agora inbox for pgsql-hackers@postgresql.org  
help / color / mirror / Atom feed
[PATCH v13 1/8] Row pattern recognition patch for raw parser.
327+ messages / 2 participants
[nested] [flat]

* [PATCH v13 1/8] Row pattern recognition patch for raw parser.
@ 2024-01-22 09:45  Tatsuo Ishii <ishii@postgresql.org>
  0 siblings, 0 replies; 327+ messages in thread

From: Tatsuo Ishii @ 2024-01-22 09:45 UTC (permalink / raw)

---
 src/backend/parser/gram.y       | 220 ++++++++++++++++++++++++++++++--
 src/include/nodes/parsenodes.h  |  57 +++++++++
 src/include/parser/kwlist.h     |   8 ++
 src/include/parser/parse_node.h |   1 +
 4 files changed, 275 insertions(+), 11 deletions(-)

diff --git a/src/backend/parser/gram.y b/src/backend/parser/gram.y
index 3460fea56b..84eb88ac2a 100644
--- a/src/backend/parser/gram.y
+++ b/src/backend/parser/gram.y
@@ -660,6 +660,21 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 				json_object_constructor_null_clause_opt
 				json_array_constructor_null_clause_opt
 
+%type <target>	row_pattern_measure_item
+				row_pattern_definition
+%type <node>	opt_row_pattern_common_syntax
+				opt_row_pattern_skip_to
+				row_pattern_subset_item
+				row_pattern_term
+%type <list>	opt_row_pattern_measures
+				row_pattern_measure_list
+				row_pattern_definition_list
+				opt_row_pattern_subset_clause
+				row_pattern_subset_list
+				row_pattern_subset_rhs
+				row_pattern
+%type <boolean>	opt_row_pattern_initial_or_seek
+				first_or_last
 
 /*
  * Non-keyword token types.  These are hard-wired into the "flex" lexer.
@@ -703,7 +718,7 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 	CURRENT_TIME CURRENT_TIMESTAMP CURRENT_USER CURSOR CYCLE
 
 	DATA_P DATABASE DAY_P DEALLOCATE DEC DECIMAL_P DECLARE DEFAULT DEFAULTS
-	DEFERRABLE DEFERRED DEFINER DELETE_P DELIMITER DELIMITERS DEPENDS DEPTH DESC
+	DEFERRABLE DEFERRED DEFINE DEFINER DELETE_P DELIMITER DELIMITERS DEPENDS DEPTH DESC
 	DETACH DICTIONARY DISABLE_P DISCARD DISTINCT DO DOCUMENT_P DOMAIN_P
 	DOUBLE_P DROP
 
@@ -719,7 +734,7 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 	HANDLER HAVING HEADER_P HOLD HOUR_P
 
 	IDENTITY_P IF_P ILIKE IMMEDIATE IMMUTABLE IMPLICIT_P IMPORT_P IN_P INCLUDE
-	INCLUDING INCREMENT INDENT INDEX INDEXES INHERIT INHERITS INITIALLY INLINE_P
+	INCLUDING INCREMENT INDENT INDEX INDEXES INHERIT INHERITS INITIAL INITIALLY INLINE_P
 	INNER_P INOUT INPUT_P INSENSITIVE INSERT INSTEAD INT_P INTEGER
 	INTERSECT INTERVAL INTO INVOKER IS ISNULL ISOLATION
 
@@ -732,7 +747,7 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 	LEADING LEAKPROOF LEAST LEFT LEVEL LIKE LIMIT LISTEN LOAD LOCAL
 	LOCALTIME LOCALTIMESTAMP LOCATION LOCK_P LOCKED LOGGED
 
-	MAPPING MATCH MATCHED MATERIALIZED MAXVALUE MERGE METHOD
+	MAPPING MATCH MATCHED MATERIALIZED MAXVALUE MEASURES MERGE METHOD
 	MINUTE_P MINVALUE MODE MONTH_P MOVE
 
 	NAME_P NAMES NATIONAL NATURAL NCHAR NEW NEXT NFC NFD NFKC NFKD NO NONE
@@ -744,8 +759,8 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 	ORDER ORDINALITY OTHERS OUT_P OUTER_P
 	OVER OVERLAPS OVERLAY OVERRIDING OWNED OWNER
 
-	PARALLEL PARAMETER PARSER PARTIAL PARTITION PASSING PASSWORD
-	PLACING PLANS POLICY
+	PARALLEL PARAMETER PARSER PARTIAL PARTITION PASSING PASSWORD PAST
+	PATTERN_P PERMUTE PLACING PLANS POLICY
 	POSITION PRECEDING PRECISION PRESERVE PREPARE PREPARED PRIMARY
 	PRIOR PRIVILEGES PROCEDURAL PROCEDURE PROCEDURES PROGRAM PUBLICATION
 
@@ -756,12 +771,13 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 	RESET RESTART RESTRICT RETURN RETURNING RETURNS REVOKE RIGHT ROLE ROLLBACK ROLLUP
 	ROUTINE ROUTINES ROW ROWS RULE
 
-	SAVEPOINT SCALAR SCHEMA SCHEMAS SCROLL SEARCH SECOND_P SECURITY SELECT
+	SAVEPOINT SCALAR SCHEMA SCHEMAS SCROLL SEARCH SECOND_P SECURITY SEEK SELECT
 	SEQUENCE SEQUENCES
+
 	SERIALIZABLE SERVER SESSION SESSION_USER SET SETS SETOF SHARE SHOW
 	SIMILAR SIMPLE SKIP SMALLINT SNAPSHOT SOME SQL_P STABLE STANDALONE_P
 	START STATEMENT STATISTICS STDIN STDOUT STORAGE STORED STRICT_P STRIP_P
-	SUBSCRIPTION SUBSTRING SUPPORT SYMMETRIC SYSID SYSTEM_P SYSTEM_USER
+	SUBSCRIPTION SUBSET SUBSTRING SUPPORT SYMMETRIC SYSID SYSTEM_P SYSTEM_USER
 
 	TABLE TABLES TABLESAMPLE TABLESPACE TEMP TEMPLATE TEMPORARY TEXT_P THEN
 	TIES TIME TIMESTAMP TO TRAILING TRANSACTION TRANSFORM
@@ -867,6 +883,7 @@ static Node *makeRecursiveViewSelect(char *relname, List *aliases, Node *query);
 %nonassoc	UNBOUNDED		/* ideally would have same precedence as IDENT */
 %nonassoc	IDENT PARTITION RANGE ROWS GROUPS PRECEDING FOLLOWING CUBE ROLLUP
 			SET KEYS OBJECT_P SCALAR VALUE_P WITH WITHOUT
+			MEASURES AFTER INITIAL SEEK PATTERN_P
 %left		Op OPERATOR		/* multi-character ops and user-defined operators */
 %left		'+' '-'
 %left		'*' '/' '%'
@@ -15928,7 +15945,8 @@ over_clause: OVER window_specification
 		;
 
 window_specification: '(' opt_existing_window_name opt_partition_clause
-						opt_sort_clause opt_frame_clause ')'
+						opt_sort_clause opt_row_pattern_measures opt_frame_clause
+						opt_row_pattern_common_syntax ')'
 				{
 					WindowDef  *n = makeNode(WindowDef);
 
@@ -15936,10 +15954,12 @@ window_specification: '(' opt_existing_window_name opt_partition_clause
 					n->refname = $2;
 					n->partitionClause = $3;
 					n->orderClause = $4;
+					n->rowPatternMeasures = $5;
 					/* copy relevant fields of opt_frame_clause */
-					n->frameOptions = $5->frameOptions;
-					n->startOffset = $5->startOffset;
-					n->endOffset = $5->endOffset;
+					n->frameOptions = $6->frameOptions;
+					n->startOffset = $6->startOffset;
+					n->endOffset = $6->endOffset;
+					n->rpCommonSyntax = (RPCommonSyntax *)$7;
 					n->location = @1;
 					$$ = n;
 				}
@@ -15963,6 +15983,31 @@ opt_partition_clause: PARTITION BY expr_list		{ $$ = $3; }
 			| /*EMPTY*/								{ $$ = NIL; }
 		;
 
+/*
+ * ROW PATTERN_P MEASURES
+ */
+opt_row_pattern_measures: MEASURES row_pattern_measure_list	{ $$ = $2; }
+			| /*EMPTY*/								{ $$ = NIL; }
+		;
+
+row_pattern_measure_list:
+			row_pattern_measure_item
+					{ $$ = list_make1($1); }
+			| row_pattern_measure_list ',' row_pattern_measure_item
+					{ $$ = lappend($1, $3); }
+		;
+
+row_pattern_measure_item:
+			a_expr AS ColLabel
+				{
+					$$ = makeNode(ResTarget);
+					$$->name = $3;
+					$$->indirection = NIL;
+					$$->val = (Node *) $1;
+					$$->location = @1;
+				}
+		;
+
 /*
  * For frame clauses, we return a WindowDef, but only some fields are used:
  * frameOptions, startOffset, and endOffset.
@@ -16122,6 +16167,143 @@ opt_window_exclusion_clause:
 			| /*EMPTY*/				{ $$ = 0; }
 		;
 
+opt_row_pattern_common_syntax:
+opt_row_pattern_skip_to opt_row_pattern_initial_or_seek
+				PATTERN_P '(' row_pattern ')'
+				opt_row_pattern_subset_clause
+				DEFINE row_pattern_definition_list
+			{
+				RPCommonSyntax *n = makeNode(RPCommonSyntax);
+				n->rpSkipTo = ((RPCommonSyntax *)$1)->rpSkipTo;
+				n->rpSkipVariable = ((RPCommonSyntax *)$1)->rpSkipVariable;
+				n->initial = $2;
+				n->rpPatterns = $5;
+				n->rpSubsetClause = $7;
+				n->rpDefs = $9;
+				$$ = (Node *) n;
+			}
+			| /*EMPTY*/		{ $$ = NULL; }
+	;
+
+opt_row_pattern_skip_to:
+			AFTER MATCH SKIP TO NEXT ROW
+				{
+					RPCommonSyntax *n = makeNode(RPCommonSyntax);
+					n->rpSkipTo = ST_NEXT_ROW;
+					n->rpSkipVariable = NULL;
+					$$ = (Node *) n;
+			}
+			| AFTER MATCH SKIP PAST LAST_P ROW
+				{
+					RPCommonSyntax *n = makeNode(RPCommonSyntax);
+					n->rpSkipTo = ST_PAST_LAST_ROW;
+					n->rpSkipVariable = NULL;
+					$$ = (Node *) n;
+				}
+			| AFTER MATCH SKIP TO first_or_last ColId
+				{
+					RPCommonSyntax *n = makeNode(RPCommonSyntax);
+					n->rpSkipTo = $5? ST_FIRST_VARIABLE : ST_LAST_VARIABLE;
+					n->rpSkipVariable = $6;
+					$$ = (Node *) n;
+				}
+/*
+			| AFTER MATCH SKIP TO LAST_P ColId		%prec LAST_P
+				{
+					RPCommonSyntax *n = makeNode(RPCommonSyntax);
+					n->rpSkipTo = ST_LAST_VARIABLE;
+					n->rpSkipVariable = $6;
+					$$ = n;
+				}
+			| AFTER MATCH SKIP TO ColId
+				{
+					RPCommonSyntax *n = makeNode(RPCommonSyntax);
+					n->rpSkipTo = ST_VARIABLE;
+					n->rpSkipVariable = $5;
+					$$ = n;
+				}
+*/
+			| /*EMPTY*/
+				{
+					RPCommonSyntax *n = makeNode(RPCommonSyntax);
+					/* temporary set default to ST_NEXT_ROW */
+					n->rpSkipTo = ST_PAST_LAST_ROW;
+					n->rpSkipVariable = NULL;
+					$$ = (Node *) n;
+				}
+	;
+
+first_or_last:
+			FIRST_P		{ $$ = true; }
+			| LAST_P	{ $$ = false; }
+	;
+
+opt_row_pattern_initial_or_seek:
+			INITIAL			{ $$ = true; }
+			| SEEK
+				{
+					ereport(ERROR,
+							(errcode(ERRCODE_SYNTAX_ERROR),
+							 errmsg("SEEK is not supported"),
+							 errhint("Use INITIAL."),
+							 parser_errposition(@1)));
+				}
+			| /*EMPTY*/		{ $$ = true; }
+		;
+
+row_pattern:
+			row_pattern_term							{ $$ = list_make1($1); }
+			| row_pattern row_pattern_term				{ $$ = lappend($1, $2); }
+		;
+
+row_pattern_term:
+			ColId	{ $$ = (Node *) makeSimpleA_Expr(AEXPR_OP, "", (Node *)makeString($1), NULL, @1); }
+			| ColId '*'	{ $$ = (Node *) makeSimpleA_Expr(AEXPR_OP, "*", (Node *)makeString($1), NULL, @1); }
+			| ColId '+'	{ $$ = (Node *) makeSimpleA_Expr(AEXPR_OP, "+", (Node *)makeString($1), NULL, @1); }
+			| ColId '?'	{ $$ = (Node *) makeSimpleA_Expr(AEXPR_OP, "?", (Node *)makeString($1), NULL, @1); }
+		;
+
+opt_row_pattern_subset_clause:
+			SUBSET row_pattern_subset_list	{ $$ = $2; }
+			| /*EMPTY*/												{ $$ = NIL; }
+		;
+
+row_pattern_subset_list:
+			row_pattern_subset_item									{ $$ = list_make1($1); }
+			| row_pattern_subset_list ',' row_pattern_subset_item	{ $$ = lappend($1, $3); }
+			| /*EMPTY*/												{ $$ = NIL; }
+		;
+
+row_pattern_subset_item: ColId '=' '(' row_pattern_subset_rhs ')'
+			{
+				RPSubsetItem *n = makeNode(RPSubsetItem);
+				n->name = $1;
+				n->rhsVariable = $4;
+				$$ = (Node *) n;
+			}
+		;
+
+row_pattern_subset_rhs:
+			ColId								{ $$ = list_make1(makeStringConst($1, @1)); }
+			| row_pattern_subset_rhs ',' ColId	{ $$ = lappend($1, makeStringConst($3, @1)); }
+			| /*EMPTY*/							{ $$ = NIL; }
+		;
+
+row_pattern_definition_list:
+			row_pattern_definition										{ $$ = list_make1($1); }
+			| row_pattern_definition_list ',' row_pattern_definition	{ $$ = lappend($1, $3); }
+		;
+
+row_pattern_definition:
+			ColId AS a_expr
+				{
+					$$ = makeNode(ResTarget);
+					$$->name = $1;
+					$$->indirection = NIL;
+					$$->val = (Node *) $3;
+					$$->location = @1;
+				}
+		;
 
 /*
  * Supporting nonterminals for expressions.
@@ -17250,6 +17432,7 @@ unreserved_keyword:
 			| INDEXES
 			| INHERIT
 			| INHERITS
+			| INITIAL
 			| INLINE_P
 			| INPUT_P
 			| INSENSITIVE
@@ -17277,6 +17460,7 @@ unreserved_keyword:
 			| MATCHED
 			| MATERIALIZED
 			| MAXVALUE
+			| MEASURES
 			| MERGE
 			| METHOD
 			| MINUTE_P
@@ -17319,6 +17503,9 @@ unreserved_keyword:
 			| PARTITION
 			| PASSING
 			| PASSWORD
+			| PAST
+			| PATTERN_P
+			| PERMUTE
 			| PLANS
 			| POLICY
 			| PRECEDING
@@ -17369,6 +17556,7 @@ unreserved_keyword:
 			| SEARCH
 			| SECOND_P
 			| SECURITY
+			| SEEK
 			| SEQUENCE
 			| SEQUENCES
 			| SERIALIZABLE
@@ -17394,6 +17582,7 @@ unreserved_keyword:
 			| STRICT_P
 			| STRIP_P
 			| SUBSCRIPTION
+			| SUBSET
 			| SUPPORT
 			| SYSID
 			| SYSTEM_P
@@ -17581,6 +17770,7 @@ reserved_keyword:
 			| CURRENT_USER
 			| DEFAULT
 			| DEFERRABLE
+			| DEFINE
 			| DESC
 			| DISTINCT
 			| DO
@@ -17743,6 +17933,7 @@ bare_label_keyword:
 			| DEFAULTS
 			| DEFERRABLE
 			| DEFERRED
+			| DEFINE
 			| DEFINER
 			| DELETE_P
 			| DELIMITER
@@ -17818,6 +18009,7 @@ bare_label_keyword:
 			| INDEXES
 			| INHERIT
 			| INHERITS
+			| INITIAL
 			| INITIALLY
 			| INLINE_P
 			| INNER_P
@@ -17867,6 +18059,7 @@ bare_label_keyword:
 			| MATCHED
 			| MATERIALIZED
 			| MAXVALUE
+			| MEASURES
 			| MERGE
 			| METHOD
 			| MINVALUE
@@ -17920,6 +18113,9 @@ bare_label_keyword:
 			| PARTITION
 			| PASSING
 			| PASSWORD
+			| PAST
+			| PATTERN_P
+			| PERMUTE
 			| PLACING
 			| PLANS
 			| POLICY
@@ -17976,6 +18172,7 @@ bare_label_keyword:
 			| SCROLL
 			| SEARCH
 			| SECURITY
+			| SEEK
 			| SELECT
 			| SEQUENCE
 			| SEQUENCES
@@ -18007,6 +18204,7 @@ bare_label_keyword:
 			| STRICT_P
 			| STRIP_P
 			| SUBSCRIPTION
+			| SUBSET
 			| SUBSTRING
 			| SUPPORT
 			| SYMMETRIC
diff --git a/src/include/nodes/parsenodes.h b/src/include/nodes/parsenodes.h
index b3181f34ae..64e9df0a48 100644
--- a/src/include/nodes/parsenodes.h
+++ b/src/include/nodes/parsenodes.h
@@ -540,6 +540,44 @@ typedef struct SortBy
 	int			location;		/* operator location, or -1 if none/unknown */
 } SortBy;
 
+/*
+ * AFTER MATCH row pattern skip to types in row pattern common syntax
+ */
+typedef enum RPSkipTo
+{
+	ST_NONE,					/* AFTER MATCH omitted */
+	ST_NEXT_ROW,				/* SKIP TO NEXT ROW */
+	ST_PAST_LAST_ROW,			/* SKIP TO PAST LAST ROW */
+	ST_FIRST_VARIABLE,			/* SKIP TO FIRST variable name */
+	ST_LAST_VARIABLE,			/* SKIP TO LAST variable name */
+	ST_VARIABLE					/* SKIP TO variable name */
+} RPSkipTo;
+
+/*
+ * Row Pattern SUBSET clause item
+ */
+typedef struct RPSubsetItem
+{
+	NodeTag		type;
+	char	   *name;			/* Row Pattern SUBSET clause variable name */
+	List	   *rhsVariable;	/* Row Pattern SUBSET rhs variables (list of char *string) */
+} RPSubsetItem;
+
+/*
+ * RowPatternCommonSyntax - raw representation of row pattern common syntax
+ *
+ */
+typedef struct RPCommonSyntax
+{
+	NodeTag		type;
+	RPSkipTo	rpSkipTo;		/* Row Pattern AFTER MATCH SKIP type */
+	char	   *rpSkipVariable;	/* Row Pattern Skip To variable name, if any */
+	bool		initial;		/* true if <row pattern initial or seek> is initial */
+	List	   *rpPatterns;		/* PATTERN variables (list of A_Expr) */
+	List	   *rpSubsetClause;	/* row pattern subset clause (list of RPSubsetItem), if any */
+	List	   *rpDefs;			/* row pattern definitions clause (list of ResTarget) */
+} RPCommonSyntax;
+
 /*
  * WindowDef - raw representation of WINDOW and OVER clauses
  *
@@ -555,6 +593,8 @@ typedef struct WindowDef
 	char	   *refname;		/* referenced window name, if any */
 	List	   *partitionClause;	/* PARTITION BY expression list */
 	List	   *orderClause;	/* ORDER BY (list of SortBy) */
+	List	   *rowPatternMeasures;	/* row pattern measures (list of ResTarget) */
+	RPCommonSyntax *rpCommonSyntax;	/* row pattern common syntax */
 	int			frameOptions;	/* frame_clause options, see below */
 	Node	   *startOffset;	/* expression for starting bound, if any */
 	Node	   *endOffset;		/* expression for ending bound, if any */
@@ -1476,6 +1516,11 @@ typedef struct GroupingSet
  * the orderClause might or might not be copied (see copiedOrder); the framing
  * options are never copied, per spec.
  *
+ * "defineClause" is Row Pattern Recognition DEFINE clause (list of
+ * TargetEntry). TargetEntry.resname represents row pattern definition
+ * variable name. "patternVariable" and "patternRegexp" represents PATTERN
+ * clause.
+ *
  * The information relevant for the query jumbling is the partition clause
  * type and its bounds.
  */
@@ -1507,6 +1552,18 @@ typedef struct WindowClause
 	Index		winref;			/* ID referenced by window functions */
 	/* did we copy orderClause from refname? */
 	bool		copiedOrder pg_node_attr(query_jumble_ignore);
+	/* Row Pattern AFTER MACH SKIP clause */
+	RPSkipTo	rpSkipTo;		/* Row Pattern Skip To type */
+	char        *rpSkipVariable;/* Row Pattern Skip To variable */
+	bool		initial;		/* true if <row pattern initial or seek> is initial */
+	/* Row Pattern DEFINE clause (list of TargetEntry) */
+	List		*defineClause;
+	/* Row Pattern DEFINE variable initial names (list of String) */
+	List		*defineInitial;
+	/* Row Pattern PATTERN variable name (list of String) */
+	List		*patternVariable;
+	/* Row Pattern PATTERN regular expression quantifier ('+' or ''. list of String) */
+	List		*patternRegexp;
 } WindowClause;
 
 /*
diff --git a/src/include/parser/kwlist.h b/src/include/parser/kwlist.h
index 2331acac09..6b3734a865 100644
--- a/src/include/parser/kwlist.h
+++ b/src/include/parser/kwlist.h
@@ -128,6 +128,7 @@ PG_KEYWORD("default", DEFAULT, RESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("defaults", DEFAULTS, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("deferrable", DEFERRABLE, RESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("deferred", DEFERRED, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("define", DEFINE, RESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("definer", DEFINER, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("delete", DELETE_P, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("delimiter", DELIMITER, UNRESERVED_KEYWORD, BARE_LABEL)
@@ -212,6 +213,7 @@ PG_KEYWORD("index", INDEX, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("indexes", INDEXES, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("inherit", INHERIT, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("inherits", INHERITS, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("initial", INITIAL, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("initially", INITIALLY, RESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("inline", INLINE_P, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("inner", INNER_P, TYPE_FUNC_NAME_KEYWORD, BARE_LABEL)
@@ -265,6 +267,7 @@ PG_KEYWORD("match", MATCH, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("matched", MATCHED, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("materialized", MATERIALIZED, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("maxvalue", MAXVALUE, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("measures", MEASURES, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("merge", MERGE, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("method", METHOD, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("minute", MINUTE_P, UNRESERVED_KEYWORD, AS_LABEL)
@@ -326,6 +329,9 @@ PG_KEYWORD("partial", PARTIAL, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("partition", PARTITION, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("passing", PASSING, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("password", PASSWORD, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("past", PAST, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("pattern", PATTERN_P, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("permute", PERMUTE, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("placing", PLACING, RESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("plans", PLANS, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("policy", POLICY, UNRESERVED_KEYWORD, BARE_LABEL)
@@ -385,6 +391,7 @@ PG_KEYWORD("scroll", SCROLL, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("search", SEARCH, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("second", SECOND_P, UNRESERVED_KEYWORD, AS_LABEL)
 PG_KEYWORD("security", SECURITY, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("seek", SEEK, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("select", SELECT, RESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("sequence", SEQUENCE, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("sequences", SEQUENCES, UNRESERVED_KEYWORD, BARE_LABEL)
@@ -416,6 +423,7 @@ PG_KEYWORD("stored", STORED, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("strict", STRICT_P, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("strip", STRIP_P, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("subscription", SUBSCRIPTION, UNRESERVED_KEYWORD, BARE_LABEL)
+PG_KEYWORD("subset", SUBSET, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("substring", SUBSTRING, COL_NAME_KEYWORD, BARE_LABEL)
 PG_KEYWORD("support", SUPPORT, UNRESERVED_KEYWORD, BARE_LABEL)
 PG_KEYWORD("symmetric", SYMMETRIC, RESERVED_KEYWORD, BARE_LABEL)
diff --git a/src/include/parser/parse_node.h b/src/include/parser/parse_node.h
index 99d6515736..1e43cef6f4 100644
--- a/src/include/parser/parse_node.h
+++ b/src/include/parser/parse_node.h
@@ -51,6 +51,7 @@ typedef enum ParseExprKind
 	EXPR_KIND_WINDOW_FRAME_RANGE,	/* window frame clause with RANGE */
 	EXPR_KIND_WINDOW_FRAME_ROWS,	/* window frame clause with ROWS */
 	EXPR_KIND_WINDOW_FRAME_GROUPS,	/* window frame clause with GROUPS */
+	EXPR_KIND_RPR_DEFINE,		/* DEFINE */
 	EXPR_KIND_SELECT_TARGET,	/* SELECT target list item */
 	EXPR_KIND_INSERT_TARGET,	/* INSERT target list item */
 	EXPR_KIND_UPDATE_SOURCE,	/* UPDATE assignment source item */
-- 
2.25.1


----Next_Part(Mon_Jan_22_19_26_18_2024_011)--
Content-Type: Text/X-Patch; charset=us-ascii
Content-Transfer-Encoding: 7bit
Content-Disposition: inline;
 filename="v13-0002-Row-pattern-recognition-patch-parse-analysis.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread

* [PATCH v44 07/10] Split cluster.h to create repack_internal.h
@ 2026-03-24 14:09  Álvaro Herrera <alvherre@kurilemu.de>
  0 siblings, 0 replies; 327+ messages in thread

From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw)

Most of the details of concurrent repack are only needed by
pgoutput_repack; and they have a nasty effect on headers included by
cluster.h (used by several high-profile places), so isolate that for
cleanliness.

Also, change_useless_for_repack() is better declared in decode.h.
---
 src/backend/commands/cluster.c                | 128 ++++++++++--------
 src/backend/postmaster/bgworker.c             |   2 +-
 .../pgoutput_repack/pgoutput_repack.c         |  13 +-
 src/include/commands/cluster.h                |  57 +-------
 src/include/commands/repack_internal.h        |  59 ++++++++
 src/include/replication/decode.h              |   4 +
 6 files changed, 147 insertions(+), 116 deletions(-)
 create mode 100644 src/include/commands/repack_internal.h

diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c
index 75556cbdafb..2c3058ba10d 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/cluster.c
@@ -59,6 +59,7 @@
 #include "commands/cluster.h"
 #include "commands/defrem.h"
 #include "commands/progress.h"
+#include "commands/repack_internal.h"
 #include "commands/tablecmds.h"
 #include "commands/vacuum.h"
 #include "executor/executor.h"
@@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared
 	char		error_queue[FLEXIBLE_ARRAY_MEMBER];
 } DecodingWorkerShared;
 
-/*
- * Generate worker's output file name. If relations of the same 'relid' happen
- * to be processed at the same time, they must be from different databases and
- * therefore different backends must be involved. (PID is already present in
- * the fileset name.)
- */
-static inline void
-DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
-{
-	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
-}
+/* Is this process a REPACK worker? */
+static bool is_repack_worker = false;
+
+static pid_t backend_pid;
+static ProcNumber backend_proc_number;
 
 /*
  * Backend-local information to control the decoding worker.
@@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL;
  */
 volatile sig_atomic_t RepackMessagePending = false;
 
+static LOCKMODE RepackLockLevel(bool concurrent);
 static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap,
 								Oid indexOid, Oid userid, LOCKMODE lmode,
 								int options);
@@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt,
 										ClusterParams *params);
 static Oid	determine_clustered_index(Relation rel, bool usingindex,
 									  const char *indexname);
-static void start_decoding_worker(Oid relid);
-static void stop_decoding_worker(void);
+
+static void start_repack_decoding_worker(Oid relid);
+static void stop_repack_decoding_worker(void);
 static void repack_worker_internal(dsm_segment *seg);
 static void export_initial_snapshot(Snapshot snapshot,
 									DecodingWorkerShared *shared);
 static Snapshot get_initial_snapshot(DecodingWorker *worker);
+extern bool am_decoding_for_repack(void);
+static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq);
+
 static void ProcessRepackMessage(StringInfo msg);
 static const char *RepackCommandAsString(RepackCommand cmd);
 
@@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 					parser_errposition(pstate, opt->location));
 	}
 
-	/*
-	 * Determine the lock mode expected by cluster_rel().
-	 *
-	 * In the exclusive case, we obtain AccessExclusiveLock right away to
-	 * avoid lock-upgrade hazard in the single-transaction case. In the
-	 * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end
-	 * of processing, supposedly for very short time. Until then, we'll have
-	 * to unlock the relation temporarily, so there's no lock-upgrade hazard.
-	 */
-	lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ?
-		AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0);
 
 	/*
 	 * If a single relation is specified, process it and we're done ... unless
@@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 										   "Repack",
 										   ALLOCSET_DEFAULT_SIZES);
 
+	/*
+	 * Since we open a new transaction for each relation, we have to check
+	 * that the relation still is what we think it is.
+	 *
+	 * In single-transaction CLUSTER, we don't need the overhead.
+	 */
 	params.options |= CLUOPT_RECHECK;
 
 	/*
@@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel)
 	MemoryContextDelete(repack_context);
 }
 
+/*
+ * In the non-concurrent case, we obtain AccessExclusiveLock throughout the
+ * operation to avoid any lock-upgrade hazards.  In the concurrent case, we
+ * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the
+ * processing and only acquire AccessExclusiveLock at the end, to swap the
+ * relation -- supposedly for a short time.
+ */
+static LOCKMODE
+RepackLockLevel(bool concurrent)
+{
+	if (concurrent)
+		return ShareUpdateExclusiveLock;
+	else
+		return AccessExclusiveLock;
+}
+
 /*
  * cluster_rel
  *
@@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	bool		concurrent = ((params->options & CLUOPT_CONCURRENT) != 0);
 	Oid			ident_idx = InvalidOid;
 
-	/*
-	 * The lock mode is AccessExclusiveLock for normal processing and
-	 * ShareUpdateExclusiveLock for concurrent processing (so that SELECT,
-	 * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be
-	 * called concurrently for the same relation).
-	 */
-	lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock;
+	/* Determine the lock mode to use. */
+	lmode = RepackLockLevel(concurrent);
 
-	/* There are specific requirements on concurrent processing. */
+	/*
+	 * Check some preconditions in the concurrent case.  This also obtains the
+	 * replica index OID.
+	 */
 	if (concurrent)
 	{
 		/*
-		 * Make sure we have no XID assigned, otherwise call of
-		 * repack_setup_logical_decoding() can cause a deadlock.
+		 * Make sure we're not in a transaction block.
 		 *
-		 * The existence of transaction block actually does not imply that XID
-		 * was already assigned, but it very likely is. We might want to check
-		 * the result of GetCurrentTransactionIdIfAny() instead, but that
-		 * would be less clear from user's perspective.
+		 * The reason is that repack_setup_logical_decoding() could deadlock
+		 * if there's an XID already assigned.  It would be possible to run in
+		 * a transaction block if we had no XID, but this restriction is
+		 * simpler for users to understand and we don't lose anything.
 		 */
 		PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)");
 
@@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 	RestrictSearchPath();
 
 	/*
-	 * Since we may open a new transaction for each relation, we have to check
-	 * that the relation still is what we think it is.
+	 * Recheck that the relation is still what it was when we started.
 	 *
-	 * If this is a single-transaction CLUSTER, we can skip these tests. We
-	 * *must* skip the one on indisclustered since it would reject an attempt
-	 * to cluster a not-previously-clustered index.
-	 *
-	 * XXX move [some of] these comments to where the RECHECK flag is
-	 * determined?
+	 * Note that it's critical to skip this in single-relation CLUSTER;
+	 * otherwise, we would reject an attempt to cluster using a
+	 * not-previously-clustered index.
 	 */
 	if (recheck &&
 		!cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid,
@@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid,
 			 * However it still seems a good practice to make sure that the
 			 * worker never survives the REPACK command.
 			 */
-			stop_decoding_worker();
+			stop_repack_decoding_worker();
 		}
 	}
 	PG_END_TRY();
@@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose,
 		 * clustering index) and checking again if it's still eligible for
 		 * REPACK CONCURRENTLY.
 		 */
-		start_decoding_worker(tableOid);
+		start_repack_decoding_worker(tableOid);
 
 		/*
 		 * Wait until the worker has the initial snapshot and retrieve it.
@@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd)
 	return "???";				/* keep compiler quiet */
 }
 
-
 /*
  * Is this backend performing logical decoding on behalf of REPACK
  * (CONCURRENTLY) ?
@@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes)
  * contents to a new table.
  */
 static void
-start_decoding_worker(Oid relid)
+start_repack_decoding_worker(Oid relid)
 {
 	Size		size;
 	dsm_segment *seg;
@@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid)
  * we need to stop it explicitly at least on ERROR in the launching backend.
  */
 static void
-stop_decoding_worker(void)
+stop_repack_decoding_worker(void)
 {
 	BgwHandleStatus status;
 
@@ -3817,12 +3822,6 @@ stop_decoding_worker(void)
 	decoding_worker = NULL;
 }
 
-/* Is this process a REPACK worker? */
-static bool is_repack_worker = false;
-
-static pid_t backend_pid;
-static ProcNumber backend_proc_number;
-
 /*
  * See ParallelWorkerShutdown for details.
  */
@@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg)
 
 	/*
 	 * Join locking group - see the comments around the call of
-	 * start_decoding_worker().
+	 * start_repack_decoding_worker().
 	 */
 	if (!BecomeLockGroupMember(shared->backend_proc, backend_pid))
 		/* The leader is not running anymore. */
@@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker)
 	return snapshot;
 }
 
+/*
+ * Generate worker's output file name. If relations of the same 'relid' happen
+ * to be processed at the same time, they must be from different databases and
+ * therefore different backends must be involved. (PID is already present in
+ * the fileset name.)
+ */
+static void
+DecodingWorkerFileName(char *fname, Oid relid, uint32 seq)
+{
+	snprintf(fname, MAXPGPATH, "%u-%u", relid, seq);
+}
+
 bool
 IsRepackWorker(void)
 {
@@ -4067,7 +4078,6 @@ void
 ProcessRepackMessages(void)
 {
 	MemoryContext oldcontext;
-
 	static MemoryContext hpm_context = NULL;
 
 	/*
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index f8a8d1681e9..9e876d55e27 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "libpq/pqsignal.h"
 #include "miscadmin.h"
 #include "pgstat.h"
diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
index 032fbd0e5b0..cc9ce615b18 100644
--- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c
+++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c
@@ -13,7 +13,7 @@
 #include "postgres.h"
 
 #include "access/detoast.h"
-#include "commands/cluster.h"
+#include "commands/repack_internal.h"
 #include "replication/snapbuild.h"
 #include "utils/memutils.h"
 
@@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 					ConcurrentChangeKind kind, HeapTuple tuple)
 {
 	RepackDecodingState *dstate;
-	MemoryContext	oldcxt;
+	MemoryContext oldcxt;
 	BufFile    *file;
 	List	   *attrs_ext = NIL;
 	int			natt_ext;
@@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 
 			slot_getsomeattrs(slot, i + 1);
 
-			/* This is a non-null varlena datum, but we only care if it's out-of-line */
+			/*
+			 * This is a non-null varlena datum, but we only care if it's
+			 * out-of-line
+			 */
 			varlen = (varlena *) DatumGetPointer(slot->tts_values[i]);
 			if (!VARATT_IS_EXTERNAL(varlen))
 				continue;
@@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation,
 				 * attributes (those actually should never appear on disk), so
 				 * only TOASTed attribute can be seen here.
 				 *
-				 * FIXME in what circumstances can an ONDISK attr appear?
-				 * Why aren't these written separately?
+				 * FIXME in what circumstances can an ONDISK attr appear? Why
+				 * aren't these written separately?
 				 */
 				Assert(VARATT_IS_EXTERNAL_ONDISK(varlen));
 			}
diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h
index 1c0ac3ab4f5..1528d34fa42 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/cluster.h
@@ -13,17 +13,12 @@
 #ifndef CLUSTER_H
 #define CLUSTER_H
 
-#include "nodes/execnodes.h"
+#include <signal.h>
+
 #include "nodes/parsenodes.h"
 #include "parser/parse_node.h"
-#include "replication/decode.h"
-#include "postmaster/bgworker.h"
-#include "replication/logical.h"
-#include "storage/buffile.h"
 #include "storage/lockdefs.h"
-#include "storage/shm_mq.h"
 #include "utils/relcache.h"
-#include "utils/resowner.h"
 
 
 /* flag bits for ClusterParams->options */
@@ -34,55 +29,14 @@
 #define CLUOPT_ANALYZE 0x08		/* do an ANALYZE */
 #define CLUOPT_CONCURRENT 0x10	/* allow concurrent data changes */
 
-
 /* options for CLUSTER */
 typedef struct ClusterParams
 {
 	bits32		options;		/* bitmask of CLUOPT_* */
 } ClusterParams;
 
-
-/*
- * The following definitions are used by REPACK CONCURRENTLY.
- */
-
-/*
- * Stored as a single byte in the output file.
- */
-#define		CHANGE_INSERT		'i'
-#define		CHANGE_UPDATE_OLD	'u'
-#define		CHANGE_UPDATE_NEW	'U'
-#define		CHANGE_DELETE		'd'
-typedef char ConcurrentChangeKind;
-
-/*
- * Logical decoding state.
- *
- * The output plugin uses it to store the data changes that it decodes from
- * WAL while the table contents is being copied to a new storage.
- */
-typedef struct RepackDecodingState
-{
-#ifdef	USE_ASSERT_CHECKING
-	/* The relation whose changes we're decoding. */
-	Oid			relid;
-#endif
-
-	/* Per-change memory context. */
-	MemoryContext	change_cxt;
-
-	/* A tuple slot used to pass tuples back and forth */
-	TupleTableSlot	*slot;
-
-	/* The current output file. */
-	BufFile    *file;
-} RepackDecodingState;
-
 extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending;
 
-extern bool IsRepackWorker(void);
-extern void HandleRepackMessageInterrupt(void);
-extern void ProcessRepackMessages(void);
 
 extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel);
 
@@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap,
 							 MultiXactId cutoffMulti,
 							 char newrelpersistence);
 
-extern bool am_decoding_for_repack(void);
-extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+extern bool IsRepackWorker(void);
+extern void HandleRepackMessageInterrupt(void);
+extern void ProcessRepackMessages(void);
+
 
-extern void RepackWorkerMain(Datum main_arg);
 #endif							/* CLUSTER_H */
diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h
new file mode 100644
index 00000000000..f90c973f5a0
--- /dev/null
+++ b/src/include/commands/repack_internal.h
@@ -0,0 +1,59 @@
+/*-------------------------------------------------------------------------
+ *
+ * repack_internal.h
+ *	  header for REPACK internals
+ *
+ * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
+ * Portions Copyright (c) 1994-5, Regents of the University of California
+ *
+ * src/include/commands/repack_internal.h
+ *
+ *-------------------------------------------------------------------------
+ */
+#ifndef REPACK_INTERNAL_H
+#define REPACK_INTERNAL_H
+
+#include "nodes/execnodes.h"
+#include "replication/decode.h"
+#include "postmaster/bgworker.h"
+#include "replication/logical.h"
+#include "storage/buffile.h"
+#include "storage/shm_mq.h"
+#include "utils/resowner.h"
+
+/*
+ * Stored as a single byte in the output file.
+ */
+#define		CHANGE_INSERT		'i'
+#define		CHANGE_UPDATE_OLD	'u'
+#define		CHANGE_UPDATE_NEW	'U'
+#define		CHANGE_DELETE		'd'
+typedef char ConcurrentChangeKind;
+
+/*
+ * Logical decoding state.
+ *
+ * The output plugin uses it to store the data changes that it decodes from
+ * WAL while the table contents is being copied to a new storage.
+ */
+typedef struct RepackDecodingState
+{
+#ifdef	USE_ASSERT_CHECKING
+	/* The relation whose changes we're decoding. */
+	Oid			relid;
+#endif
+
+	/* Per-change memory context. */
+	MemoryContext change_cxt;
+
+	/* A tuple slot used to pass tuples back and forth */
+	TupleTableSlot *slot;
+
+	/* The current output file. */
+	BufFile    *file;
+} RepackDecodingState;
+
+
+extern void RepackWorkerMain(Datum main_arg);
+
+#endif							/* REPACK_INTERNAL_H */
diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h
index 49f00fc48b8..02b5393474c 100644
--- a/src/include/replication/decode.h
+++ b/src/include/replication/decode.h
@@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf
 extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx,
 										 XLogReaderState *record);
 
+/* in commands/cluster.c */
+extern bool change_useless_for_repack(XLogRecordBuffer *buf);
+
+
 #endif
-- 
2.47.3


--gwom7bl7ogtszo4k
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
	filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch"



^ permalink  raw  reply  [nested|flat] 327+ messages in thread


end of thread, other threads:[~2026-03-24 14:09 UTC | newest]

Thread overview: 327+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2024-01-22 09:45 [PATCH v13 1/8] Row pattern recognition patch for raw parser. Tatsuo Ishii <ishii@postgresql.org>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox